# -*- coding: utf-8 -*-
"""
Web crawler for intangible heritage sports university inheritance data from the China Intangible Cultural Heritage Digital Museum (ihchina.cn)

Functions:
1. Read the excel file "清洗后_仅新增非遗体育主项目清单.xlsx" and extract cleaned‑up main‑item names as matching keywords
2. Fetch the list of "participating universities of the inheritance training plan" and "intangible heritage into campus" related news from ihchina.cn
3. Extract university names, intangible heritage projects and course information mentioned in the texts
4. Export search results to an excel file in unified format

Data sources:
- 2024-2025 annual list of participating universities of China Intangible Cultural Heritage Inheritance Training Plan (130 universities)
- News and activity reports related to intangible heritage entering campus

Note: the website has no public JSON interface, data is obtained via HTML parsing.
"""

import pandas as pd
import time
import re
import os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

# ========================== File Path Configuration ==========================
INPUT_KEYWORDS = "清洗后_仅新增非遗体育主项目清单.xlsx"
OUTPUT_FILE = "检索结果6_中国非遗数字博物馆高校传承数据.xlsx"

# ========================== HTTP Request Header Configuration ==========================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
    "Referer": "https://www.ihchina.cn/",
}

# ========================== Data Reading Module ==========================
def read_keywords():
    """Read and return the cleaned‑up main‑item keyword list"""
    if not os.path.exists(INPUT_KEYWORDS):
        raise FileNotFoundError(f"Input file not found: {INPUT_KEYWORDS}")

    df = pd.read_excel(INPUT_KEYWORDS)
    if "清洗后主项目名称" not in df.columns:
        raise ValueError(f"Required column missing in excel. Current column list: {df.columns.tolist()}")

    keywords = df["清洗后主项目名称"].dropna().astype(str).str.strip().tolist()
    keywords = [k for k in keywords if k and k != "nan"]

    if not keywords:
        raise ValueError("No valid matching keywords extracted")

    print(f"✅ Successfully loaded {len(keywords)} matching keywords")
    return keywords


# ========================== HTTP Request with Automatic Retry ==========================
def fetch_html(url, max_retries=2, timeout=15):
    """Send an HTTP GET request with automatic retry on failure"""
    for attempt in range(max_retries + 1):
        try:
            response = requests.get(url, headers=HEADERS, timeout=timeout)
            response.encoding = "utf-8"
            if response.status_code == 200:
                return response.text
            else:
                print(f"  ⚠️ HTTP {response.status_code} (Attempt {attempt+1}/{max_retries+1}): {url}")
        except requests.exceptions.RequestException as e:
            print(f"  ⚠️ Request Exception (Attempt {attempt+1}/{max_retries+1}): {str(e)[:50]}")
        if attempt < max_retries:
            time.sleep(2)
    return None


# ========================== Extract University Names ==========================
def extract_institutions(text):
    """
    Extract university names from text
    Matching pattern: XX大学, XX学院, XX职业技术大学, etc.
    """
    patterns = [
        r'([\u4e00-\u9fa5]{2,20}(?:大学|学院|职业技术大学|职业学院|师范高等专科学校|高等专科学校|体育学院|音乐学院|艺术学院|美术学院|戏曲学院|舞蹈学院|工艺美术高级技工学校))',
        r'([\u4e00-\u9fa5]{2,10}(?:科技|理工|工业|农业|林业|财经|工商|交通|民族|师范|医药|艺术|体育|音乐|美术|戏曲|舞蹈)大学)',
        r'([\u4e00-\u9fa5]{2,10}(?:科技|理工|工业|农业|林业|财经|工商|交通|民族|师范|医药|艺术|体育|音乐|美术|戏曲|舞蹈)学院)',
    ]
    institutions = []
    for pattern in patterns:
        matches = re.findall(pattern, text)
        for m in matches:
            if len(m) >= 4 and m not in institutions:
                institutions.append(m)
    return institutions


def infer_university_type(name):
    """Infer university type from its name"""
    sports_keywords = ["体育", "运动", "武术"]
    normal_keywords = ["师范", "教育"]
    ethnic_keywords = ["民族"]

    if any(k in name for k in sports_keywords):
        return "体育类"
    if any(k in name for k in normal_keywords):
        return "师范类"
    if any(k in name for k in ethnic_keywords):
        return "民族类"
    if any(k in name for k in ["大学", "学院"]):
        return "综合类"
    return "综合类"


def infer_location(name):
    """Simply infer location from the name prefix"""
    province_map = {
        "北京": "北京", "上海": "上海", "天津": "天津", "重庆": "重庆",
        "河北": "河北", "山西": "山西", "辽宁": "辽宁", "吉林": "吉林",
        "黑龙江": "黑龙江", "江苏": "江苏", "浙江": "浙江", "安徽": "安徽",
        "福建": "福建", "江西": "江西", "山东": "山东", "河南": "河南",
        "湖北": "湖北", "湖南": "湖南", "广东": "广东", "海南": "海南",
        "四川": "四川", "贵州": "贵州", "云南": "云南", "陕西": "陕西",
        "甘肃": "甘肃", "青海": "青海", "台湾": "台湾",
        "内蒙古": "内蒙古", "广西": "广西", "西藏": "西藏",
        "宁夏": "宁夏", "新疆": "新疆",
    }
    for key in province_map:
        if name.startswith(key):
            return province_map[key]
    return ""


# ========================== Match Intangible Heritage Sports Projects ==========================
def match_sports_keywords(text, keywords):
    """
    Match intangible heritage sports project keywords within the text
    Return the list of matched keywords
    """
    matched = []
    for kw in keywords:
        if kw in text:
            matched.append(kw)
    return matched


# ========================== Fetch the School List of the Training Plan ==========================
def fetch_yanpei_schools():
    """
    Fetch the list of participating universities of the intangible heritage training plan from ihchina.cn
    Return: list of university names
    """
    print("\n📚 Fetching the participating university list of the 'Intangible Heritage Inheritance Training Plan'...")

    # Training plan list page
    url = "https://www.ihchina.cn/art/detail/id/28645.html"
    html = fetch_html(url)

    if not html:
        print("  ❌ Failed to fetch the training plan list page")
        return []

    soup = BeautifulSoup(html, "html.parser")
    content = soup.get_text()

    # Extract all university names (using regex)
    schools = extract_institutions(content)

    # Deduplicate
    schools = list(dict.fromkeys(schools))

    print(f"  ✅ Extracted {len(schools)} universities from the training plan list")
    return schools


# ========================== Fetch the News about Intangible Heritage Entering Campus ==========================
def fetch_news_pages():
    """
    Fetch news pages related to 'intangible heritage entering campus' from ihchina.cn
    Return: list of news page URLs
    """
    print("\n📚 Fetching news related to 'intangible heritage entering campus'...")

    # Search related news (via site search function)
    search_urls = [
        "https://www.ihchina.cn/search?keyword=非遗进校园&page=1",
        "https://www.ihchina.cn/search?keyword=高校传承&page=1",
        "https://www.ihchina.cn/search?keyword=研培计划&page=1",
        "https://www.ihchina.cn/search?keyword=非遗课程&page=1",
    ]

    news_urls = []
    for search_url in search_urls:
        html = fetch_html(search_url)
        if not html:
            continue
        soup = BeautifulSoup(html, "html.parser")
        # Find news links
        for a in soup.find_all("a", href=True):
            href = a["href"]
            if "/news" in href or "/project" in href:
                full_url = urljoin("https://www.ihchina.cn", href)
                if full_url not in news_urls:
                    news_urls.append(full_url)
        time.sleep(3)  # Request interval

    # Deduplicate
    news_urls = list(dict.fromkeys(news_urls))
    print(f"  ✅ Found {len(news_urls)} related news pages")
    return news_urls[:30]  # Limit to 30 pages at most to avoid excessive requests


def parse_news_page(url, keywords):
    """
    Parse a single news page and extract university and course information
    Return: [(university name, matched intangible heritage project list, course description), ...]
    """
    html = fetch_html(url)
    if not html:
        return []

    soup = BeautifulSoup(html, "html.parser")
    content = soup.get_text()

    # Extract universities
    institutions = extract_institutions(content)

    results = []
    for inst in institutions:
        # Match intangible heritage sports projects in the current page
        matched_projects = match_sports_keywords(content, keywords)
        if matched_projects:
            # Extract course related description (context containing the matched keyword)
            course_desc = ""
            for kw in matched_projects:
                idx = content.find(kw)
                if idx != -1:
                    start = max(0, idx - 30)
                    end = min(len(content), idx + len(kw) + 80)
                    snippet = content[start:end].replace("\n", " ").strip()
                    if len(snippet) > 10:
                        course_desc = snippet
                        break

            for project in matched_projects:
                results.append({
                    "institution": inst,
                    "project": project,
                    "course_desc": course_desc
                })

    return results


# ========================== Main Program Entry ==========================
def main():
    print("=" * 70)
    print("China Intangible Cultural Heritage Digital Museum Intangible Heritage Sports University Inheritance Data Crawler")
    print("Data source: intangible heritage training plan list + news about intangible heritage entering campus")
    print("Note: the website has no public JSON interface, data is obtained via HTML parsing")
    print("=" * 70)

    try:
        # ---------- Step 1: Load matching keywords ----------
        keywords = read_keywords()
        print(f"Keyword preview (first 10 items): {keywords[:10]}")

        # ---------- Step 2: Fetch training plan universities ----------
        yanpei_schools = fetch_yanpei_schools()

        # ---------- Step 3: Fetch news about intangible heritage entering campus ----------
        news_urls = fetch_news_pages()

        # ---------- Step 4: Parse news pages ----------
        print("\n" + "=" * 70)
        print("🔍 Start parsing news pages and extracting university and intangible heritage project information")
        print("=" * 70)

        all_results = []
        total_news = len(news_urls)

        for idx, url in enumerate(news_urls, 1):
            print(f"\n[{idx}/{total_news}] Parsing news: {url[:80]}...")

            news_results = parse_news_page(url, keywords)

            if news_results:
                # Deduplicate (same university + same project)
                unique_pairs = set()
                for item in news_results:
                    key = (item["institution"], item["project"])
                    if key not in unique_pairs:
                        unique_pairs.add(key)
                        all_results.append(item)
                print(f"  ✅ Extracted {len(news_results)} records (newly added after deduplication: {len(set((r['institution'], r['project']) for r in news_results))})")
            else:
                print(f"  ⚠️ No intangible heritage sports projects matched")

            # Request interval ≥ 3 seconds
            if idx < total_news:
                time.sleep(3)

        # ---------- Step 5: Use training plan universities as supplementary source ----------
        # For universities in the training plan, no record is generated if no specific project is matched in news (no specific project info available)
        # But if already matched in news, it is already included

        # ---------- Step 6: Build output data ----------
        print("\n" + "=" * 70)
        print("📊 Building output data")
        print("=" * 70)

        output_records = []
        for item in all_results:
            inst = item["institution"]
            project = item["project"]

            # Infer university type and location
            univ_type = infer_university_type(inst)
            location = infer_location(inst)

            # Course description
            course_desc = item.get("course_desc", "")
            if not course_desc:
                course_desc = f"该院校在非遗进校园相关活动中涉及「{project}」项目"

            output_records.append({
                "对应非遗体育项目名称": project,
                "院校名称": inst,
                "院校所在地": location,
                "院校类型": univ_type,
                "非遗传统体育课程名称": project,
                "课程属性": "公共选修课",  # According to MOE requirements, intangible heritage inheritance bases should open public elective courses[reference:5]
                "课程等级": "本科",
                "课程简介": course_desc[:200] + "..." if len(course_desc) > 200 else course_desc,
                "数据来源渠道": "中国非物质文化遗产数字博物馆",
                "备注": "非遗研培/校园传承项目"
            })

        # ---------- Step 7: Deduplicate ----------
        if output_records:
            df_results = pd.DataFrame(output_records)
            # Deduplicate by university name + intangible heritage sports course name
            df_results = df_results.drop_duplicates(
                subset=["院校名称", "非遗传统体育课程名称"],
                keep="first"
            )
        else:
            df_results = pd.DataFrame(columns=[
                "对应非遗体育项目名称", "院校名称", "院校所在地", "院校类型",
                "非遗传统体育课程名称", "课程属性", "课程等级", "课程简介",
                "数据来源渠道", "备注"
            ])

        # ---------- Step 8: Export to excel ----------
        df_results.to_excel(OUTPUT_FILE, index=False)

        # ---------- Step 9: Print statistical report ----------
        print("\n" + "=" * 70)
        print("📊 Crawling‑task Completion Statistics")
        print("=" * 70)
        print(f"Total matching keywords: {len(keywords)}")
        print(f"Training plan university count: {len(yanpei_schools)}")
        print(f"Parsed news page count: {len(news_urls)}")
        print(f"Matched records (before deduplication): {len(output_records)}")
        print(f"Deduplicated records: {len(df_results)}")
        print(f"\n✅ Output file: {OUTPUT_FILE}")
        print("=" * 70)

        # Preview result data
        if not df_results.empty:
            print("\n📖 Result preview (first 10 entries):")
            print("-" * 70)
            preview_cols = ["对应非遗体育项目名称", "院校名称", "非遗传统体育课程名称"]
            print(df_results[preview_cols].head(10).to_string(index=False))
            print("-" * 70)

            # Count matched universities per project
            print("\n📊 Number of matched universities per intangible heritage sports project:")
            course_count = df_results.groupby("对应非遗体育项目名称").size().sort_values(ascending=False)
            for project, count in course_count.head(15).items():
                print(f"  {project}: {count} universities")
            if len(course_count) > 15:
                print(f"  ... {len(course_count) - 15} more projects omitted")
        else:
            print("\n⚠️ No data matched")
            print("Potential reasons:")
            print("1. No news about these intangible heritage sports projects entering campus exists on ihchina.cn")
            print("2. Keywords and news content are expressed inconsistently")
            print("3. Website structure changed, parsing rules need updating")

        # ---------- Step 10: Data source description ----------
        print("\n" + "=" * 70)
        print("📌 Data Source Description")
        print("=" * 70)
        print("The data comes from the China Intangible Cultural Heritage Digital Museum (ihchina.cn):")
        print("  1. 2024-2025 annual list of participating universities of China Intangible Cultural Heritage Inheritance Training Plan (130 universities)[reference:6]")
        print("  2. News and activity reports related to 'intangible heritage entering campus'")
        print("\n💡 Notes:")
        print("  - The website publishes content as static HTML, with no public JSON interface")
        print("  - Participating universities of the training plan all offer intangible heritage related courses or research programs[reference:7]")
        print("  - '课程属性' is labeled as '公共选修课' according to MOE requirements[reference:8]")
        print("  - All records are labeled as '非遗研培/校园传承项目'")
        print("=" * 70)

    except FileNotFoundError as e:
        print(f"❌ File‑related error: {e}")
    except ValueError as e:
        print(f"❌ Data‑validation error: {e}")
    except Exception as e:
        print(f"❌ Unexpected runtime error: {e}")
        import traceback
        traceback.print_exc()


if __name__ == "__main__":
    main()
